Détection de Malware avec Machine Learning 🧠🔍

Ce projet consiste à utiliser des algorithmes de machine learning pour analyser des fichiers et détecter des comportements malveillants. Le modèle est entraîné sur des jeux de données de fichiers malveillants et légitimes afin d'identifier des signatures de malware.

Fonctionnalités du Projet :

Pré-requis :

Script Python pour la Détection de Malware :


import os
import pickle
import logging
import numpy as np
from sklearn.ensemble import RandomForestClassifier
from sklearn.model_selection import train_test_split
from sklearn.metrics import classification_report, confusion_matrix, roc_curve, auc
import matplotlib.pyplot as plt

# Configuration de la journalisation
logging.basicConfig(filename="malware_detection.log", level=logging.INFO, format="%(asctime)s - %(message)s")

# Extraction des caractéristiques
def extract_features(filepath):
    try:
        size = os.path.getsize(filepath)
        permissions = oct(os.stat(filepath).st_mode)[-3:]
        entropy = calculate_entropy(filepath)
        sha256 = calculate_sha256(filepath)
        return [size, int(permissions), entropy]
    except Exception as e:
        logging.error(f"Erreur d'extraction pour {filepath}: {e}")
        return None

# Calcul de l'entropie
def calculate_entropy(filepath):
    with open(filepath, 'rb') as f:
        data = f.read()
    if not data:
        return 0
    entropy = -sum(p * np.log2(p) for p in np.bincount(data, minlength=256) / len(data) if p > 0)
    return entropy

# Calcul du checksum SHA256
import hashlib
def calculate_sha256(filepath):
    with open(filepath, 'rb') as f:
        return hashlib.sha256(f.read()).hexdigest()

# Entraînement du modèle
def train_model(X, y):
    model = RandomForestClassifier(n_estimators=100, random_state=42)
    model.fit(X, y)
    return model

# Évaluation et courbes
def evaluate_model(y_test, y_pred):
    print("Matrice de confusion :")
    print(confusion_matrix(y_test, y_pred))
    fpr, tpr, _ = roc_curve(y_test, y_pred)
    roc_auc = auc(fpr, tpr)
    plt.figure()
    plt.plot(fpr, tpr, label=f"Courbe ROC (AUC = {roc_auc:.2f})")
    plt.xlabel("Faux positifs")
    plt.ylabel("Vrais positifs")
    plt.title("Courbe ROC")
    plt.legend()
    plt.show()

# Main training function
def main_train(train_data):
    X, y = [], []
    for filepath, label in train_data:
        features = extract_features(filepath)
        if features is not None:
            X.append(features)
            y.append(label)

    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)
    model = train_model(X_train, y_train)
    with open("malware_model.pkl", "wb") as f:
        pickle.dump(model, f)

    y_pred = model.predict(X_test)
    print(classification_report(y_test, y_pred))
    evaluate_model(y_test, y_pred)

# Utilisation du modèle pour classifier
def classify_file(filepath):
    try:
        with open("malware_model.pkl", "rb") as f:
            model = pickle.load(f)
        features = extract_features(filepath)
        if features:
            result = model.predict([features])[0]
            return "Malware" if result == 1 else "Légitime"
        else:
            return "Erreur de classification"
    except FileNotFoundError:
        return "Modèle non trouvé. Veuillez entraîner le modèle avant de classifier."

if __name__ == "__main__":
    # Exemple d'utilisation
    train_data = [("file_legit_1.exe", 0), ("file_malware_1.exe", 1)]
    main_train(train_data)
    result = classify_file("unknown_file.exe")
    print(f"Résultat : {result}")

Instructions d'Utilisation :

🚀 Entraînement :

Pour entraîner le modèle, fournissez un jeu de données et exécutez :

python detect_malware.py --train chemin_vers_donnees

📋 Classification :

Pour classifier un fichier inconnu :

python detect_malware.py --classify chemin_vers_fichier

Attention et Éthique :

Utilisez ce projet uniquement dans un cadre légal. L’analyse de fichiers sans autorisation est strictement interdite. Respectez les lois en vigueur et utilisez cet outil de manière éthique. 🛡️